Launch Options in verl-omni Example Scripts
Briefly introduces the launch options in verl-omni’s example scripts, as some of them are related to issues like OOM and algorithm tuning. This post first takes LTX2.3 FlowGRPO training script as an example.
python3 -m verl_omni.trainer.main_diffusion \
trainer.use_v1=false \
trainer.device=npu \
data.train_files=$train_path \
data.val_files=$test_path \
data.train_batch_size=32 \
data.val_max_samples=1024 \
data.max_prompt_length=1024 \
data.truncation=error \
data.seed=42 \
algorithm.adv_estimator=flow_grpo \
algorithm.global_std=True \
actor_rollout_ref.model.path=$MODEL_PATH \
actor_rollout_ref.model.algorithm=flow_grpo \
actor_rollout_ref.model.attn_backend=native \
actor_rollout_ref.model.enable_gradient_checkpointing=True \
actor_rollout_ref.model.lora_rank=64 \
actor_rollout_ref.model.lora_alpha=128 \
actor_rollout_ref.model.target_modules="$ltx_lora_targets" \
actor_rollout_ref.model.fsdp_layer_prefixes="['transformer_blocks.']" \
actor_rollout_ref.actor.strategy=fsdp \
'+actor_rollout_ref.actor.fsdp_config.wrap_policy.transformer_layer_cls_to_wrap=[LTX2VideoTransformerBlock]' \
actor_rollout_ref.actor.optim.lr=3e-4 \
actor_rollout_ref.actor.optim.weight_decay=1e-4 \
actor_rollout_ref.actor.optim.betas="[0.9,0.999]" \
actor_rollout_ref.actor.optim.override_optimizer_config="{eps: 1e-8}" \
actor_rollout_ref.actor.optim.clip_grad=1.0 \
actor_rollout_ref.actor.ppo_mini_batch_size=16 \
actor_rollout_ref.actor.ppo_micro_batch_size_per_gpu=8 \
actor_rollout_ref.actor.diffusion_loss.clip_ratio=1e-4 \
actor_rollout_ref.actor.diffusion_loss.adv_clip_max=5.0 \
actor_rollout_ref.actor.use_kl_loss=False \
actor_rollout_ref.actor.fsdp_config.model_dtype=bfloat16 \
actor_rollout_ref.actor.fsdp_config.param_offload=True \
actor_rollout_ref.actor.fsdp_config.optimizer_offload=True \
actor_rollout_ref.actor.fsdp_config.ulysses_sequence_parallel_size=1 \
actor_rollout_ref.rollout.name=vllm_omni \
actor_rollout_ref.rollout.rollout_attn_backend=TORCH_SDPA \
actor_rollout_ref.rollout.tensor_model_parallel_size=$ROLLOUT_TP \
actor_rollout_ref.rollout.n=8 \
actor_rollout_ref.rollout.seed=42 \
actor_rollout_ref.rollout.agent.num_workers=$((NUM_GPUS / ROLLOUT_TP)) \
actor_rollout_ref.rollout.agent.default_agent_loop=ltx2_diffusion_single_turn_agent \
actor_rollout_ref.rollout.load_format=safetensors \
actor_rollout_ref.rollout.layered_summon=True \
actor_rollout_ref.rollout.log_prob_micro_batch_size_per_gpu=16 \
actor_rollout_ref.rollout.pipeline.height=256 \
actor_rollout_ref.rollout.pipeline.width=384 \
actor_rollout_ref.rollout.pipeline.num_frames=81 \
actor_rollout_ref.rollout.pipeline.frame_rate=24.0 \
actor_rollout_ref.rollout.pipeline.num_inference_steps=24 \
actor_rollout_ref.rollout.pipeline.guidance_scale=4.0 \
actor_rollout_ref.rollout.pipeline.max_sequence_length=1024 \
+actor_rollout_ref.rollout.pipeline.output_type=pt \
actor_rollout_ref.rollout.algo.noise_level=0.8 \
actor_rollout_ref.rollout.algo.sde_type=cps \
actor_rollout_ref.rollout.algo.sde_window_range="[0,10]" \
actor_rollout_ref.rollout.algo.sde_window_size=3 \
actor_rollout_ref.rollout.algo.sde_contiguous=False \
actor_rollout_ref.rollout.algo.sde_window_seed=42 \
actor_rollout_ref.rollout.calculate_log_probs=True \
actor_rollout_ref.rollout.val_kwargs.pipeline.height=256 \
actor_rollout_ref.rollout.val_kwargs.pipeline.width=384 \
actor_rollout_ref.rollout.val_kwargs.pipeline.num_frames=81 \
actor_rollout_ref.rollout.val_kwargs.pipeline.frame_rate=24.0 \
actor_rollout_ref.rollout.val_kwargs.pipeline.num_inference_steps=50 \
actor_rollout_ref.rollout.val_kwargs.pipeline.guidance_scale=4.0 \
+actor_rollout_ref.rollout.val_kwargs.pipeline.output_type=pt \
actor_rollout_ref.rollout.val_kwargs.algo.noise_level=0.0 \
actor_rollout_ref.ref.log_prob_micro_batch_size_per_gpu=16 \
reward.num_workers=$REWARD_NUM_WORKERS \
reward.reward_model.enable=False \
reward.custom_reward_function.path=pkg://verl_omni.reward_loop.reward_manager.multi \
reward.custom_reward_function.name=_multi_reward_placeholder \
reward.reward_manager.name=MultiVisualRewardManager \
reward.reward_manager.module.path=pkg://verl_omni.reward_loop.reward_manager \
"+reward.reward_functions.clap.path=$repo_root/verl_omni/utils/reward_score/clap.py" \
'+reward.reward_functions.clap.name=compute_score' \
'+reward.reward_functions.clap.weight=1.0' \
"+reward.reward_functions.clap.device=$REWARD_DEVICE:0" \
"+reward.reward_functions.clap.model_name_or_path=$CLAP_MODEL_PATH" \
"+reward.reward_functions.imagebind.path=$repo_root/verl_omni/utils/reward_score/imagebind.py" \
'+reward.reward_functions.imagebind.name=compute_score' \
'+reward.reward_functions.imagebind.weight=1.0' \
"+reward.reward_functions.imagebind.device=$REWARD_DEVICE:1" \
"+reward.reward_functions.imagebind.model_name_or_path=$IMAGEBIND_MODEL_PATH" \
'+reward.reward_functions.imagebind.mode=audio_video' \
reward.aggregation=weighted_sum \
trainer.logger='["console","tensorboard","wandb"]' \
trainer.project_name=flow_grpo_npu \
trainer.experiment_name=ltx2_3_t2av_lora_npu \
trainer.default_local_dir=$checkpoint_dir \
trainer.resume_mode=disable \
trainer.log_val_generations=8 \
trainer.video_fps=24 \
trainer.val_before_train=True \
trainer.n_gpus_per_node=$NUM_GPUS \
trainer.nnodes=1 \
trainer.save_freq=50 \
trainer.test_freq=20 \
trainer.total_epochs=15 \
trainer.total_training_steps=$TOTAL_TRAINING_STEPS "$@"